agora inbox for [email protected]help / color / mirror / Atom feed
[PATCH v28 09/11] Add support for min/max aggregates for IVM 385+ messages / 2 participants [nested] [flat]
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v32 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index ecec93ec1c..d3cf3df5a7 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1311,6 +1311,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index b7f6c3831b..42ff0ed98f 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -105,6 +133,7 @@ typedef struct MV_TriggerTable TupleTableSlot *slot; /* for checking visibility in the pre-state table */ } MV_TriggerTable; +static HTAB *mv_query_cache = NULL; static HTAB *mv_trigger_info = NULL; static bool in_delta_calculation = false; @@ -165,6 +194,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -173,17 +205,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort); /* @@ -2124,6 +2169,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2205,6 +2252,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2234,6 +2292,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2252,10 +2312,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2447,6 +2515,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2549,19 +2681,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2576,10 +2733,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2587,10 +2745,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2773,6 +2946,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -2806,6 +3322,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -2814,6 +3337,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.25.1 --Multipart=_Sun__31_Mar_2024_22_59_31_+0900_msknEviJj08_wgqO Content-Type: text/x-diff; name="v32-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Disposition: attachment; filename="v32-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Transfer-Encoding: 7bit ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v34 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index d8767137d9..abce06d046 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1309,6 +1309,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index ca6d236989..8cd7ed3b68 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -105,6 +133,7 @@ typedef struct MV_TriggerTable TupleTableSlot *slot; /* for checking visibility in the pre-state table */ } MV_TriggerTable; +static HTAB *mv_query_cache = NULL; static HTAB *mv_trigger_info = NULL; static bool in_delta_calculation = false; @@ -165,6 +194,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -173,17 +205,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort); /* @@ -2124,6 +2169,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2205,6 +2252,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2234,6 +2292,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2252,10 +2312,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2447,6 +2515,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2549,19 +2681,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2576,10 +2733,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2587,10 +2745,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2773,6 +2946,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -2806,6 +3322,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -2814,6 +3337,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.34.1 --Multipart=_Thu__11_Jul_2024_13_23_57_+0900_hKsU2G3a3BgS2FFs Content-Type: text/x-diff; name="v34-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v34-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v28 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index c8aa558f2e..c40ea6b2bc 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1312,6 +1312,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index aa6bf2694a..fa65e4aeb0 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -73,6 +73,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + OverrideSearchPath *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -109,6 +137,7 @@ typedef struct MV_TriggerTable TupleTableSlot *slot; /* for checking visibility in the pre-state table */ } MV_TriggerTable; +static HTAB *mv_query_cache = NULL; static HTAB *mv_trigger_info = NULL; static bool in_delta_calculation = false; @@ -169,6 +198,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -177,17 +209,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort); /* @@ -2102,6 +2147,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2183,6 +2230,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2212,6 +2270,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2230,10 +2290,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2425,6 +2493,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2527,19 +2659,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2554,10 +2711,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2565,10 +2723,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2751,6 +2924,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -2784,6 +3300,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -2792,6 +3315,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + OverrideSearchPathMatchesCurrent(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetOverrideSearchPath(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.25.1 --Multipart=_Thu__1_Jun_2023_23_59_09_+0900_/G5+8nG46.f1T42K Content-Type: text/x-diff; name="v28-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Disposition: attachment; filename="v28-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Transfer-Encoding: 7bit ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v29 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index c8aa558f2e..c40ea6b2bc 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1312,6 +1312,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index ee41f0007d..eff512d40c 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -73,6 +73,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -109,6 +137,7 @@ typedef struct MV_TriggerTable TupleTableSlot *slot; /* for checking visibility in the pre-state table */ } MV_TriggerTable; +static HTAB *mv_query_cache = NULL; static HTAB *mv_trigger_info = NULL; static bool in_delta_calculation = false; @@ -169,6 +198,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -177,17 +209,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort); /* @@ -2101,6 +2146,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2182,6 +2229,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2211,6 +2269,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2229,10 +2289,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2424,6 +2492,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2526,19 +2658,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2553,10 +2710,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2564,10 +2722,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2750,6 +2923,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -2783,6 +3299,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -2791,6 +3314,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.25.1 --Multipart=_Mon__28_Aug_2023_11_52_52_+0900_hj6L5h176QaSGtg7 Content-Type: text/x-diff; name="v29-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Disposition: attachment; filename="v29-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Transfer-Encoding: 7bit ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v33 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index d8767137d9..abce06d046 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1309,6 +1309,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 97406b28c9..0ff5e3922b 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -105,6 +133,7 @@ typedef struct MV_TriggerTable TupleTableSlot *slot; /* for checking visibility in the pre-state table */ } MV_TriggerTable; +static HTAB *mv_query_cache = NULL; static HTAB *mv_trigger_info = NULL; static bool in_delta_calculation = false; @@ -165,6 +194,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -173,17 +205,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort); /* @@ -2122,6 +2167,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2203,6 +2250,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2232,6 +2290,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2250,10 +2310,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2445,6 +2513,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2547,19 +2679,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2574,10 +2731,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2585,10 +2743,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2771,6 +2944,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -2804,6 +3320,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -2812,6 +3335,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.25.1 --Multipart=_Tue__2_Jul_2024_17_03_11_+0900_6OrVBZxOB4o_k6A1 Content-Type: text/x-diff; name="v33-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v33-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v30 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index d93eec3eec..0536e44b9e 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1312,6 +1312,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 3c523991ed..e60f92226c 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -73,6 +73,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -109,6 +137,7 @@ typedef struct MV_TriggerTable TupleTableSlot *slot; /* for checking visibility in the pre-state table */ } MV_TriggerTable; +static HTAB *mv_query_cache = NULL; static HTAB *mv_trigger_info = NULL; static bool in_delta_calculation = false; @@ -169,6 +198,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -177,17 +209,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort); /* @@ -2123,6 +2168,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2204,6 +2251,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2233,6 +2291,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2251,10 +2311,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2446,6 +2514,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2548,19 +2680,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2575,10 +2732,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2586,10 +2744,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2772,6 +2945,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -2805,6 +3321,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -2813,6 +3336,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.25.1 --Multipart=_Mon__4_Mar_2024_11_58_46_+0900_UaponF/qQhQrVCFt Content-Type: text/x-diff; name="v30-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Disposition: attachment; filename="v30-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Transfer-Encoding: 7bit ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v31 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index aa8440b4e1..7e1627ad1b 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1309,6 +1309,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 97406b28c9..0ff5e3922b 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -105,6 +133,7 @@ typedef struct MV_TriggerTable TupleTableSlot *slot; /* for checking visibility in the pre-state table */ } MV_TriggerTable; +static HTAB *mv_query_cache = NULL; static HTAB *mv_trigger_info = NULL; static bool in_delta_calculation = false; @@ -165,6 +194,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -173,17 +205,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort); /* @@ -2122,6 +2167,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2203,6 +2250,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2232,6 +2290,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2250,10 +2310,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2445,6 +2513,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2547,19 +2679,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2574,10 +2731,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2585,10 +2743,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2771,6 +2944,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -2804,6 +3320,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -2812,6 +3335,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.25.1 --Multipart=_Fri__29_Mar_2024_23_47_00_+0900_KGpmmDOIs1266Ib1 Content-Type: text/x-diff; name="v31-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Disposition: attachment; filename="v31-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Transfer-Encoding: 7bit ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v29 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index c8aa558f2e..c40ea6b2bc 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1312,6 +1312,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index ee41f0007d..eff512d40c 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -73,6 +73,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -109,6 +137,7 @@ typedef struct MV_TriggerTable TupleTableSlot *slot; /* for checking visibility in the pre-state table */ } MV_TriggerTable; +static HTAB *mv_query_cache = NULL; static HTAB *mv_trigger_info = NULL; static bool in_delta_calculation = false; @@ -169,6 +198,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -177,17 +209,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort); /* @@ -2101,6 +2146,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2182,6 +2229,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2211,6 +2269,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2229,10 +2289,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2424,6 +2492,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2526,19 +2658,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2553,10 +2710,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2564,10 +2722,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2750,6 +2923,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -2783,6 +3299,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -2791,6 +3314,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.25.1 --Multipart=_Mon__28_Aug_2023_16_05_30_+0900_b1OvQD_3A3ZMTGvj Content-Type: text/x-diff; name="v29-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Disposition: attachment; filename="v29-0010-Add-regression-tests-for-Incremental-View-Mainte.patch" Content-Transfer-Encoding: 7bit ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v37 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 45a30309951..ed67f399b7c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1332,6 +1332,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Fri__29_May_2026_23_14_17_+0900_Te0o73X2VqYK57Gd Content-Type: text/x-diff; name="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v37-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH v38 09/11] Add support for min/max aggregates for IVM @ 2023-05-31 11:58 Yugo Nagata <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Yugo Nagata @ 2023-05-31 11:58 UTC (permalink / raw) Supporting min and max is more complicated than count, sum, or avg. For an example of min, when tuples are inserted, the current min value in the view and the min value in the inseteted tuples are compared, then the smaller one is used as the latest min value. On the other hand, when tuples are deleted, if the current min value in the view equals to the min in the deleted tuples, we need re-computation the latest min value from base tables. Otherwise, the current value in the view remains. --- src/backend/commands/createas.c | 45 +++ src/backend/commands/matview.c | 644 +++++++++++++++++++++++++++++++- 2 files changed, 680 insertions(+), 9 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c index 35e124694ab..540d9e3d5c8 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -1341,6 +1341,51 @@ check_aggregate_supports_ivm(Oid aggfnoid) case F_AVG_FLOAT8: case F_AVG_INTERVAL: + /* min */ + case F_MIN_ANYARRAY: + case F_MIN_INT8: + case F_MIN_INT4: + case F_MIN_INT2: + case F_MIN_OID: + case F_MIN_FLOAT4: + case F_MIN_FLOAT8: + case F_MIN_DATE: + case F_MIN_TIME: + case F_MIN_TIMETZ: + case F_MIN_MONEY: + case F_MIN_TIMESTAMP: + case F_MIN_TIMESTAMPTZ: + case F_MIN_INTERVAL: + case F_MIN_TEXT: + case F_MIN_NUMERIC: + case F_MIN_BPCHAR: + case F_MIN_TID: + case F_MIN_ANYENUM: + case F_MIN_INET: + case F_MIN_PG_LSN: + + /* max */ + case F_MAX_ANYARRAY: + case F_MAX_INT8: + case F_MAX_INT4: + case F_MAX_INT2: + case F_MAX_OID: + case F_MAX_FLOAT4: + case F_MAX_FLOAT8: + case F_MAX_DATE: + case F_MAX_TIME: + case F_MAX_TIMETZ: + case F_MAX_MONEY: + case F_MAX_TIMESTAMP: + case F_MAX_TIMESTAMPTZ: + case F_MAX_INTERVAL: + case F_MAX_TEXT: + case F_MAX_NUMERIC: + case F_MAX_BPCHAR: + case F_MAX_TID: + case F_MAX_ANYENUM: + case F_MAX_INET: + case F_MAX_PG_LSN: return true; default: diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 710224aa994..3b3871c49e2 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -69,6 +69,34 @@ typedef struct #define MV_INIT_QUERYHASHSIZE 16 +/* MV query type codes */ +#define MV_PLAN_RECALC 1 +#define MV_PLAN_SET_VALUE 2 + +/* + * MI_QueryKey + * + * The key identifying a prepared SPI plan in our query hashtable + */ +typedef struct MV_QueryKey +{ + Oid matview_id; /* OID of materialized view */ + int32 query_type; /* query type ID, see MV_PLAN_XXX above */ +} MV_QueryKey; + +/* + * MV_QueryHashEntry + * + * Hash entry for cached plans used to maintain materialized views. + */ +typedef struct MV_QueryHashEntry +{ + MV_QueryKey key; + SPIPlanPtr plan; + SearchPathMatcher *search_path; /* search_path used for parsing + * and planning */ +} MV_QueryHashEntry; + /* * MV_TriggerHashEntry * @@ -121,6 +149,7 @@ typedef struct DroppedImmvInfo } DroppedImmvInfo; static HTAB *mv_trigger_info = NULL; +static HTAB *mv_query_cache = NULL; static HTAB *dropped_immv = NULL; static HTAB *LastIvmUpdateHash; @@ -210,6 +239,9 @@ static void append_set_clause_for_sum(const char *resname, StringInfo buf_old, static void append_set_clause_for_avg(const char *resname, StringInfo buf_old, StringInfo buf_new, StringInfo aggs_list, const char *aggtype); +static void append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min); static char *get_operation_string(IvmOp op, const char *col, const char *arg1, const char *arg2, const char* count_col, const char *castType); static char *get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, @@ -218,17 +250,30 @@ static void apply_old_delta(const char *matviewname, const char *deltaname_old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname); + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc); static void apply_new_delta(const char *matviewname, const char *deltaname_new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_new, List *keys, StringInfo target_list, StringInfo aggs_set, const char* count_colname); static char *get_matching_condition_string(List *keys); +static char *get_returning_string(List *minmax_list, List *is_min_list, List *keys); +static char *get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list); +static char *get_select_for_recalc_string(List *keys); +static void recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel); +static SPIPlanPtr get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes); +static SPIPlanPtr get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); static void mv_InitHashTables(void); +static SPIPlanPtr mv_FetchPreparedPlan(MV_QueryKey *key); +static void mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan); +static void mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type); static void clean_up_IVM_hash_entry(MV_TriggerHashEntry *entry, bool is_abort, SubTransactionId subxid); static void setLastUpdateXid(Oid immv_oid, FullTransactionId xid); @@ -2343,6 +2388,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n ListCell *lc; int i; List *keys = NIL; + List *minmax_list = NIL; + List *is_min_list = NIL; /* @@ -2424,6 +2471,17 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_list_buf, format_type_be(aggref->aggtype)); + /* min/max */ + else if (!strcmp(aggname, "min") || !strcmp(aggname, "max")) + { + bool is_min = (!strcmp(aggname, "min")); + + append_set_clause_for_minmax(resname, aggs_set_old, aggs_set_new, aggs_list_buf, is_min); + + /* make a resname list of min and max aggregates */ + minmax_list = lappend(minmax_list, resname); + is_min_list = lappend_int(is_min_list, is_min); + } else elog(ERROR, "unsupported aggregate function: %s", aggname); } @@ -2453,6 +2511,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) { EphemeralNamedRelation enr = palloc(sizeof(EphemeralNamedRelationData)); + SPITupleTable *tuptable_recalc = NULL; + uint64 num_recalc; int rc; /* convert tuplestores to ENR, and register for SPI */ @@ -2471,10 +2531,18 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, Tuplestorestate *n /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, keys, aggs_list_buf, aggs_set_old, - count_colname); + minmax_list, is_min_list, + count_colname, &tuptable_recalc, &num_recalc); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); + /* + * If we have min or max, we might have to recalculate aggregate values from base tables + * on some tuples. TIDs and keys such tuples are returned as a result of the above query. + */ + if (minmax_list && tuptable_recalc) + recalc_and_set_values(tuptable_recalc, num_recalc, minmax_list, keys, matviewRel); + } /* For tuple insertion */ if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) @@ -2666,6 +2734,70 @@ append_set_clause_for_avg(const char *resname, StringInfo buf_old, ); } +/* + * append_set_clause_for_minmax + * + * Append SET clause string for min or max aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + * is_min is true if this is min, false if not. + */ +static void +append_set_clause_for_minmax(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + bool is_min) +{ + char *count_col = IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* + * If the new value doesn't became NULL then use the value remaining + * in the view although this will be recomputated afterwords. + */ + appendStringInfo(buf_old, + ", %s = CASE WHEN %s THEN NULL ELSE %s END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_SUB, "mv", "t", count_col), + quote_qualified_identifier("mv", resname) + ); + /* count = mv.count - t.count */ + appendStringInfo(buf_old, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* + * min = LEAST(mv.min, diff.min) + * max = GREATEST(mv.max, diff.max) + */ + appendStringInfo(buf_new, + ", %s = CASE WHEN %s THEN NULL ELSE %s(%s,%s) END", + quote_qualified_identifier(NULL, resname), + get_null_condition_string(IVM_ADD, "mv", "diff", count_col), + + is_min ? "LEAST" : "GREATEST", + quote_qualified_identifier("mv", resname), + quote_qualified_identifier("diff", resname) + ); + /* count = mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s = %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + /* * get_operation_string * @@ -2768,19 +2900,44 @@ get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, * list to identify a tuple in the view. If the view has aggregates, this * requires strings representing resnames of aggregates and SET clause for * updating aggregate values. + * + * If the view has min or max aggregate, this requires a list of resnames of + * min/max aggregates and a list of boolean which represents which entries in + * minmax_list is min. These are necessary to check if we need to recalculate + * min or max aggregate values. In this case, this query returns TID and keys + * of tuples which need to be recalculated. This result and the number of rows + * are stored in tuptables and num_recalc repectedly. + * */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, List *keys, StringInfo aggs_list, StringInfo aggs_set, - const char *count_colname) + List *minmax_list, List *is_min_list, + const char *count_colname, + SPITupleTable **tuptable_recalc, uint64 *num_recalc) { StringInfoData querybuf; char *match_cond; + char *updt_returning = ""; + char *select_for_recalc = "SELECT"; bool agg_without_groupby = (list_length(keys) == 0); + Assert(tuptable_recalc != NULL); + Assert(num_recalc != NULL); + /* build WHERE condition for searching tuples to be deleted */ match_cond = get_matching_condition_string(keys); + /* + * We need a special RETURNING clause and SELECT statement for min/max to + * check which tuple needs re-calculation from base tables. + */ + if (minmax_list) + { + updt_returning = get_returning_string(minmax_list, is_min_list, keys); + select_for_recalc = get_select_for_recalc_string(keys); + } + /* Search for matching tuples from the view and update or delete if found. */ initStringInfo(&querybuf); appendStringInfo(&querybuf, @@ -2795,10 +2952,11 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, "UPDATE %s AS mv SET %s = mv.%s OPERATOR(pg_catalog.-) t.%s " "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND NOT for_dlt " - ")" - /* delete a tuple if this is to be deleted */ - "DELETE FROM %s AS mv USING t " - "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt", + "%s" /* RETURNING clause for recalc infomation */ + "), dlt AS (" /* delete a tuple if this is to be deleted */ + "DELETE FROM %s AS mv USING t " + "WHERE mv.ctid OPERATOR(pg_catalog.=) t.ctid AND for_dlt" + ") %s", /* SELECT returning which tuples need to be recalculated */ count_colname, count_colname, count_colname, (agg_without_groupby ? "false" : "true"), (aggs_list != NULL ? aggs_list->data : ""), @@ -2806,10 +2964,25 @@ apply_old_delta_with_count(const char *matviewname, const char *deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, (aggs_set != NULL ? aggs_set->data : ""), - matviewname); + updt_returning, + matviewname, + select_for_recalc); - if (SPI_exec(querybuf.data, 0) != SPI_OK_DELETE) + if (SPI_exec(querybuf.data, 0) != SPI_OK_SELECT) elog(ERROR, "SPI_exec failed: %s", querybuf.data); + + + /* Return tuples to be recalculated. */ + if (minmax_list) + { + *tuptable_recalc = SPI_tuptable; + *num_recalc = SPI_processed; + } + else + { + *tuptable_recalc = NULL; + *num_recalc = 0; + } } /* @@ -2992,6 +3165,349 @@ get_matching_condition_string(List *keys) return match_cond.data; } +/* + * get_returning_string + * + * Build a string for RETURNING clause of UPDATE used in apply_old_delta_with_count. + * This clause returns ctid and a boolean value that indicates if we need to + * recalculate min or max value, for each updated row. + */ +static char * +get_returning_string(List *minmax_list, List *is_min_list, List *keys) +{ + StringInfoData returning; + char *recalc_cond; + ListCell *lc; + + Assert(minmax_list != NIL && is_min_list != NIL); + recalc_cond = get_minmax_recalc_condition_string(minmax_list, is_min_list); + + initStringInfo(&returning); + + appendStringInfo(&returning, "RETURNING mv.ctid AS tid, (%s) AS recalc", recalc_cond); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + appendStringInfo(&returning, ", %s", quote_qualified_identifier("mv", resname)); + } + + return returning.data; +} + +/* + * get_minmax_recalc_condition_string + * + * Build a predicate string for checking if any min/max aggregate + * value needs to be recalculated. + */ +static char * +get_minmax_recalc_condition_string(List *minmax_list, List *is_min_list) +{ + StringInfoData recalc_cond; + ListCell *lc1, *lc2; + + initStringInfo(&recalc_cond); + + Assert (list_length(minmax_list) == list_length(is_min_list)); + + forboth (lc1, minmax_list, lc2, is_min_list) + { + char *resname = (char *) lfirst(lc1); + bool is_min = (bool) lfirst_int(lc2); + char *op_str = (is_min ? ">=" : "<="); + + appendStringInfo(&recalc_cond, "%s OPERATOR(pg_catalog.%s) %s", + quote_qualified_identifier("mv", resname), + op_str, + quote_qualified_identifier("t", resname) + ); + + if (lnext(minmax_list, lc1)) + appendStringInfo(&recalc_cond, " OR "); + } + + return recalc_cond.data; +} + +/* + * get_select_for_recalc_string + * + * Build a query to return tid and keys of tuples which need + * recalculation. This is used as the result of the query + * built by apply_old_delta. + */ +static char * +get_select_for_recalc_string(List *keys) +{ + StringInfoData qry; + ListCell *lc; + + initStringInfo(&qry); + + appendStringInfo(&qry, "SELECT tid"); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + appendStringInfo(&qry, ", %s", NameStr(attr->attname)); + } + + appendStringInfo(&qry, " FROM updt WHERE recalc"); + + return qry.data; +} + +/* + * recalc_and_set_values + * + * Recalculate tuples in a materialized from base tables and update these. + * The tuples which needs recalculation are specified by keys, and resnames + * of columns to be updated are specified by namelist. TIDs and key values + * are given by tuples in tuptable_recalc. Its first attribute must be TID + * and key values must be following this. + */ +static void +recalc_and_set_values(SPITupleTable *tuptable_recalc, int64 num_tuples, + List *namelist, List *keys, Relation matviewRel) +{ + TupleDesc tupdesc_recalc = tuptable_recalc->tupdesc; + Oid *keyTypes = NULL, *types = NULL; + char *keyNulls = NULL, *nulls = NULL; + Datum *keyVals = NULL, *vals = NULL; + int num_vals = list_length(namelist); + int num_keys = list_length(keys); + uint64 i; + Oid matviewOid; + char *matviewname; + + matviewOid = RelationGetRelid(matviewRel); + matviewname = quote_qualified_identifier(get_namespace_name(RelationGetNamespace(matviewRel)), + RelationGetRelationName(matviewRel)); + + /* If we have keys, initialize arrays for them. */ + if (keys) + { + keyTypes = palloc(sizeof(Oid) * num_keys); + keyNulls = palloc(sizeof(char) * num_keys); + keyVals = palloc(sizeof(Datum) * num_keys); + /* a tuple contains keys to be recalculated and ctid to be updated*/ + Assert(tupdesc_recalc->natts == num_keys + 1); + + /* Types of key attributes */ + for (i = 0; i < num_keys; i++) + keyTypes[i] = TupleDescAttr(tupdesc_recalc, i + 1)->atttypid; + } + + /* allocate memory for all attribute names and tid */ + types = palloc(sizeof(Oid) * (num_vals + 1)); + nulls = palloc(sizeof(char) * (num_vals + 1)); + vals = palloc(sizeof(Datum) * (num_vals + 1)); + + /* For each tuple which needs recalculation */ + for (i = 0; i < num_tuples; i++) + { + int j; + bool isnull; + SPIPlanPtr plan; + SPITupleTable *tuptable_newvals; + TupleDesc tupdesc_newvals; + + /* Set group key values as parameters if needed. */ + if (keys) + { + for (j = 0; j < num_keys; j++) + { + keyVals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, j + 2, &isnull); + if (isnull) + keyNulls[j] = 'n'; + else + keyNulls[j] = ' '; + } + } + + /* + * Get recalculated values from base tables. The result must be + * only one tuple thich contains the new values for specified keys. + */ + plan = get_plan_for_recalc(matviewOid, namelist, keys, keyTypes); + if (SPI_execute_plan(plan, keyVals, keyNulls, false, 0) != SPI_OK_SELECT) + elog(ERROR, "SPI_execute_plan"); + if (SPI_processed != 1) + elog(ERROR, "SPI_execute_plan returned zero or more than one rows"); + + tuptable_newvals = SPI_tuptable; + tupdesc_newvals = tuptable_newvals->tupdesc; + + Assert(tupdesc_newvals->natts == num_vals); + + /* Set the new values as parameters */ + for (j = 0; j < tupdesc_newvals->natts; j++) + { + if (i == 0) + types[j] = TupleDescAttr(tupdesc_newvals, j)->atttypid; + + vals[j] = SPI_getbinval(tuptable_newvals->vals[0], tupdesc_newvals, j + 1, &isnull); + if (isnull) + nulls[j] = 'n'; + else + nulls[j] = ' '; + } + /* Set TID of the view tuple to be updated as a parameter */ + types[j] = TIDOID; + vals[j] = SPI_getbinval(tuptable_recalc->vals[i], tupdesc_recalc, 1, &isnull); + nulls[j] = ' '; + + /* Update the view tuple to the new values */ + plan = get_plan_for_set_values(matviewOid, matviewname, namelist, types); + if (SPI_execute_plan(plan, vals, nulls, false, 0) != SPI_OK_UPDATE) + elog(ERROR, "SPI_execute_plan"); + } +} + + +/* + * get_plan_for_recalc + * + * Create or fetch a plan for recalculating value in the view's target list + * from base tables using the definition query of materialized view specified + * by matviewOid. namelist is a list of resnames of values to be recalculated. + * + * keys is a list of keys to identify tuples to be recalculated if this is not + * empty. KeyTypes is an array of types of keys. + */ +static SPIPlanPtr +get_plan_for_recalc(Oid matviewOid, List *namelist, List *keys, Oid *keyTypes) +{ + MV_QueryKey hash_key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the recalculation */ + mv_BuildQueryKey(&hash_key, matviewOid, MV_PLAN_RECALC); + if ((plan = mv_FetchPreparedPlan(&hash_key)) == NULL) + { + ListCell *lc; + StringInfoData str; + char *viewdef; + + /* get view definition of matview */ + viewdef = text_to_cstring((text *) DatumGetPointer( + DirectFunctionCall1(pg_get_viewdef, ObjectIdGetDatum(matviewOid)))); + /* get rid of trailing semi-colon */ + viewdef[strlen(viewdef)-1] = '\0'; + + /* + * Build a query string for recalculating values. This is like + * + * SELECT x1, x2, x3, ... FROM ( ... view definition query ...) mv + * WHERE (key1, key2, ...) = ($1, $2, ...); + */ + + initStringInfo(&str); + appendStringInfo(&str, "SELECT "); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, " FROM (%s) mv", viewdef); + + if (keys) + { + int i = 1; + char paramname[16]; + + appendStringInfo(&str, " WHERE ("); + foreach (lc, keys) + { + Form_pg_attribute attr = (Form_pg_attribute) lfirst(lc); + char *resname = NameStr(attr->attname); + Oid typid = attr->atttypid; + + sprintf(paramname, "$%d", i); + appendStringInfo(&str, "("); + generate_equal(&str, typid, resname, paramname); + appendStringInfo(&str, " OR (%s IS NULL AND %s IS NULL))", + resname, paramname); + + if (lnext(keys, lc)) + appendStringInfoString(&str, " AND "); + i++; + } + appendStringInfo(&str, ")"); + } + else + keyTypes = NULL; + + plan = SPI_prepare(str.data, list_length(keys), keyTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&hash_key, plan); + } + + return plan; +} + +/* + * get_plan_for_set_values + * + * Create or fetch a plan for applying new values calculated by + * get_plan_for_recalc to a materialized view specified by matviewOid. + * matviewname is the name of the view. namelist is a list of resnames + * of attributes to be updated, and valTypes is an array of types of the + * values. + */ +static SPIPlanPtr +get_plan_for_set_values(Oid matviewOid, char *matviewname, List *namelist, + Oid *valTypes) +{ + MV_QueryKey key; + SPIPlanPtr plan; + + /* Fetch or prepare a saved plan for the real check */ + mv_BuildQueryKey(&key, matviewOid, MV_PLAN_SET_VALUE); + if ((plan = mv_FetchPreparedPlan(&key)) == NULL) + { + ListCell *lc; + StringInfoData str; + int i; + + /* + * Build a query string for applying min/max values. This is like + * + * UPDATE matviewname AS mv + * SET (x1, x2, x3, x4) = ($1, $2, $3, $4) + * WHERE ctid = $5; + */ + + initStringInfo(&str); + appendStringInfo(&str, "UPDATE %s AS mv SET (", matviewname); + foreach (lc, namelist) + { + appendStringInfo(&str, "%s", (char *) lfirst(lc)); + if (lnext(namelist, lc)) + appendStringInfoString(&str, ", "); + } + appendStringInfo(&str, ") = ROW("); + + for (i = 1; i <= list_length(namelist); i++) + appendStringInfo(&str, "%s$%d", (i==1 ? "" : ", "), i); + + appendStringInfo(&str, ") WHERE ctid OPERATOR(pg_catalog.=) $%d", i); + + plan = SPI_prepare(str.data, list_length(namelist) + 1, valTypes); + if (plan == NULL) + elog(ERROR, "SPI_prepare returned %s for %s", SPI_result_code_string(SPI_result), str.data); + + SPI_keepplan(plan); + mv_HashPreparedPlan(&key, plan); + } + + return plan; +} + /* * generate_equals * @@ -3025,6 +3541,13 @@ mv_InitHashTables(void) { HASHCTL ctl; + memset(&ctl, 0, sizeof(ctl)); + ctl.keysize = sizeof(MV_QueryKey); + ctl.entrysize = sizeof(MV_QueryHashEntry); + mv_query_cache = hash_create("MV query cache", + MV_INIT_QUERYHASHSIZE, + &ctl, HASH_ELEM | HASH_BLOBS); + memset(&ctl, 0, sizeof(ctl)); ctl.keysize = sizeof(Oid); ctl.entrysize = sizeof(MV_TriggerHashEntry); @@ -3039,6 +3562,109 @@ mv_InitHashTables(void) &ctl, HASH_ELEM | HASH_BLOBS); } +/* + * mv_FetchPreparedPlan + */ +static SPIPlanPtr +mv_FetchPreparedPlan(MV_QueryKey *key) +{ + MV_QueryHashEntry *entry; + SPIPlanPtr plan; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Lookup for the key + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_FIND, NULL); + if (entry == NULL) + return NULL; + + /* + * Check whether the plan is still valid. If it isn't, we don't want to + * simply rely on plancache.c to regenerate it; rather we should start + * from scratch and rebuild the query text too. This is to cover cases + * such as table/column renames. We depend on the plancache machinery to + * detect possible invalidations, though. + * + * CAUTION: this check is only trustworthy if the caller has already + * locked both materialized views and base tables. + * + * Also, check whether the search_path is still the same as when we made it. + * If it isn't, we need to rebuild the query text because the result of + * pg_ivm_get_viewdef() will change. + */ + plan = entry->plan; + if (plan && SPI_plan_is_valid(plan) && + SearchPathMatchesCurrentEnvironment(entry->search_path)) + return plan; + + /* + * Otherwise we might as well flush the cached plan now, to free a little + * memory space before we make a new one. + */ + if (plan) + SPI_freeplan(plan); + if (entry->search_path) + pfree(entry->search_path); + + entry->plan = NULL; + entry->search_path = NULL; + + return NULL; +} + +/* + * mv_HashPreparedPlan + * + * Add another plan to our private SPI query plan hashtable. + */ +static void +mv_HashPreparedPlan(MV_QueryKey *key, SPIPlanPtr plan) +{ + MV_QueryHashEntry *entry; + bool found; + + /* + * On the first call initialize the hashtable + */ + if (!mv_query_cache) + mv_InitHashTables(); + + /* + * Add the new plan. We might be overwriting an entry previously found + * invalid by mv_FetchPreparedPlan. + */ + entry = (MV_QueryHashEntry *) hash_search(mv_query_cache, + (void *) key, + HASH_ENTER, &found); + Assert(!found || entry->plan == NULL); + entry->plan = plan; + entry->search_path = GetSearchPathMatcher(TopMemoryContext); +} + +/* + * mv_BuildQueryKey + * + * Construct a hashtable key for a prepared SPI plan for IVM. + */ +static void +mv_BuildQueryKey(MV_QueryKey *key, Oid matview_id, int32 query_type) +{ + /* + * We assume struct MV_QueryKey contains no padding bytes, else we'd need + * to use memset to clear them. + */ + key->matview_id = matview_id; + key->query_type = query_type; +} + /* * AtAbort_IVM * -- 2.43.0 --Multipart=_Wed__1_Jul_2026_00_04_01_+0900_OVSy2WWK_9aByzDJ Content-Type: text/x-diff; name="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Disposition: attachment; filename="v38-0008-Add-aggregates-support-in-IVM.patch" Content-Transfer-Encoding: quoted-printable ^ permalink raw reply [nested|flat] 385+ messages in thread
* [PATCH 5/6] Use background worker to do logical decoding. @ 2026-01-12 16:30 Antonin Houska <[email protected]> 0 siblings, 0 replies; 385+ messages in thread From: Antonin Houska @ 2026-01-12 16:30 UTC (permalink / raw) If the backend performing REPACK (CONCURRENTLY) does both data copying and logical decoding, it has to "travel in time" back and forth and therefore it has to invalidate system caches quite a few times. (The copying and the decoding work with different catalog snapshots.) As the decoding worker has separate caches, the switching is not necessary. Without the worker, it'd also be difficult to switch between potentiallly long running tasks like index build and WAL decoding. (No decoding during that time at all can suspend archiving / recycling of WAL segments for some time, which in turn may result in full disk.) Another problem is that, after having acquired AccessExclusiveLock (in order to swap the files), the backend needs to both decode and apply the data changes that took place while it was waiting for the lock. With the decoding worker, the decoding runs all the time, so the backend only needs to apply the changes. This can reduce the time the exclusive lock is held for. Note that the code added in order to handle ERRORs in the background worker almost duplicates the existing code that does the same for other types of workers (See ProcessParallelMessages() and ProcessParallelApplyMessages()). Refactoring of the existing code might be useful, to reduce the duplication. --- src/backend/access/heap/heapam_handler.c | 44 - src/backend/commands/cluster.c | 1176 +++++++++++++---- src/backend/libpq/pqmq.c | 5 + src/backend/postmaster/bgworker.c | 4 + src/backend/replication/logical/logical.c | 6 +- .../pgoutput_repack/pgoutput_repack.c | 54 +- src/backend/storage/ipc/procsignal.c | 4 + src/backend/tcop/postgres.c | 4 + .../utils/activity/wait_event_names.txt | 2 + src/include/access/tableam.h | 7 +- src/include/commands/cluster.h | 71 +- src/include/storage/procsignal.h | 1 + src/tools/pgindent/typedefs.list | 4 +- 13 files changed, 981 insertions(+), 401 deletions(-) diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c index 3526b6adcb5..475c536ce43 100644 --- a/src/backend/access/heap/heapam_handler.c +++ b/src/backend/access/heap/heapam_handler.c @@ -33,7 +33,6 @@ #include "catalog/index.h" #include "catalog/storage.h" #include "catalog/storage_xlog.h" -#include "commands/cluster.h" #include "commands/progress.h" #include "executor/executor.h" #include "miscadmin.h" @@ -688,7 +687,6 @@ heapam_relation_copy_for_cluster(Relation OldHeap, Relation NewHeap, Relation OldIndex, bool use_sort, TransactionId OldestXmin, Snapshot snapshot, - LogicalDecodingContext *decoding_ctx, TransactionId *xid_cutoff, MultiXactId *multi_cutoff, double *num_tuples, @@ -710,7 +708,6 @@ heapam_relation_copy_for_cluster(Relation OldHeap, Relation NewHeap, BufferHeapTupleTableSlot *hslot; BlockNumber prev_cblock = InvalidBlockNumber; bool concurrent = snapshot != NULL; - XLogRecPtr end_of_wal_prev = GetFlushRecPtr(NULL); /* Remember if it's a system catalog */ is_system_catalog = IsSystemRelation(OldHeap); @@ -957,31 +954,6 @@ heapam_relation_copy_for_cluster(Relation OldHeap, Relation NewHeap, ct_val[1] = *num_tuples; pgstat_progress_update_multi_param(2, ct_index, ct_val); } - - /* - * Process the WAL produced by the load, as well as by other - * transactions, so that the replication slot can advance and WAL does - * not pile up. Use wal_segment_size as a threshold so that we do not - * introduce the decoding overhead too often. - * - * Of course, we must not apply the changes until the initial load has - * completed. - * - * Note that our insertions into the new table should not be decoded - * as we (intentionally) do not write the logical decoding specific - * information to WAL. - */ - if (concurrent) - { - XLogRecPtr end_of_wal; - - end_of_wal = GetFlushRecPtr(NULL); - if ((end_of_wal - end_of_wal_prev) > wal_segment_size) - { - repack_decode_concurrent_changes(decoding_ctx, end_of_wal); - end_of_wal_prev = end_of_wal; - } - } } if (indexScan != NULL) @@ -1027,22 +999,6 @@ heapam_relation_copy_for_cluster(Relation OldHeap, Relation NewHeap, /* Report n_tuples */ pgstat_progress_update_param(PROGRESS_REPACK_HEAP_TUPLES_INSERTED, n_tuples); - - /* - * Try to keep the amount of not-yet-decoded WAL small, like - * above. - */ - if (concurrent) - { - XLogRecPtr end_of_wal; - - end_of_wal = GetFlushRecPtr(NULL); - if ((end_of_wal - end_of_wal_prev) > wal_segment_size) - { - repack_decode_concurrent_changes(decoding_ctx, end_of_wal); - end_of_wal_prev = end_of_wal; - } - } } tuplesort_end(tuplesort); diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 9f61620a279..49a42f54ae4 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -12,12 +12,13 @@ * In concurrent mode, we lock the table with only ShareUpdateExclusiveLock, * then do an initial copy as above. However, while the tuples are being * copied, concurrent transactions could modify the table. To cope with those - * changes, we rely on logical decoding to obtain them from WAL. The changes - * are accumulated in a tuplestore. Once the initial copy is complete, we - * read the changes from the tuplestore and re-apply them on the new heap. - * Then we upgrade our ShareUpdateExclusiveLock to AccessExclusiveLock and - * swap the relfilenodes. This way, the time we hold a strong lock on the - * table is much reduced, and the bloat is eliminated. + * changes, we rely on logical decoding to obtain them from WAL. A bgworker + * consumes WAL while the initial copy is ongoing (to prevent excessive WAL + * from being reserved), and accumulates the changes in a file. Once the + * initial copy is complete, we read the changes from the file and re-apply + * them on the new heap. Then we upgrade our ShareUpdateExclusiveLock to + * AccessExclusiveLock and swap the relfilenodes. This way, the time we hold + * a strong lock on the table is much reduced, and the bloat is eliminated. * * There is hardly anything left of Paul Brown's original implementation... * @@ -45,6 +46,7 @@ #include "access/xlog_internal.h" #include "access/xloginsert.h" #include "access/xlogutils.h" +#include "access/xlogwait.h" #include "catalog/catalog.h" #include "catalog/dependency.h" #include "catalog/heap.h" @@ -61,6 +63,8 @@ #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" +#include "libpq/pqformat.h" +#include "libpq/pqmq.h" #include "miscadmin.h" #include "optimizer/optimizer.h" #include "pgstat.h" @@ -71,6 +75,8 @@ #include "storage/ipc.h" #include "storage/lmgr.h" #include "storage/predicate.h" +#include "storage/procsignal.h" +#include "tcop/tcopprot.h" #include "utils/acl.h" #include "utils/fmgroids.h" #include "utils/guc.h" @@ -117,6 +123,12 @@ typedef struct IndexInsertState /* The WAL segment being decoded. */ static XLogSegNo repack_current_segment = 0; +/* + * The first file exported by the decoding worker must contain a snapshot, the + * following ones contain the data changes. + */ +#define WORKER_FILE_SNAPSHOT 0 + /* * Information needed to apply concurrent data changes. */ @@ -136,8 +148,113 @@ typedef struct ChangeDest /* Needed to update indexes of rel_dst. */ IndexInsertState *iistate; + + /* + * Sequential number of the file containing the changes. + * + * TODO This field makes the structure name less descriptive. Should we + * rename it, e.g. to ChangeApplyInfo? + */ + int file_seq; } ChangeDest; +/* + * Layout of shared memory used for communication between backend and the + * worker that performs logical decoding of data changes + */ +typedef struct DecodingWorkerShared +{ + /* Is the decoding initialized? */ + bool initialized; + + /* + * Once the worker has reached this LSN, it should close the current + * output file and either create a new one or exit, according to the field + * 'done'. If the value is InvalidXLogRecPtr, the worker should decode all + * the WAL available and keep checking this field. It is ok if the worker + * had already decoded records whose LSN is >= lsn_upto before this field + * has been set. + */ + XLogRecPtr lsn_upto; + + /* Exit after closing the current file? */ + bool done; + + /* The output is stored here. */ + SharedFileSet sfs; + + /* Number of the last file exported by the worker. */ + int last_exported; + + /* Synchronize access to the fields above. */ + slock_t mutex; + + /* Database to connect to. */ + Oid dbid; + + /* Role to connect as. */ + Oid roleid; + + /* Decode data changes of this relation. */ + Oid relid; + + /* The backend uses this to wait for the worker. */ + ConditionVariable cv; + + /* Info to signal the backend. */ + PGPROC *backend_proc; + pid_t backend_pid; + ProcNumber backend_proc_number; + + /* Error queue. */ + shm_mq *error_mq; + + /* + * Memory the queue is located int. + * + * For considerations on the value see the comments of + * PARALLEL_ERROR_QUEUE_SIZE. + */ +#define REPACK_ERROR_QUEUE_SIZE 16384 + char error_queue[FLEXIBLE_ARRAY_MEMBER]; +} DecodingWorkerShared; + +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static inline void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + +/* + * Backend-local information to control the decoding worker. + */ +typedef struct DecodingWorker +{ + /* The worker. */ + BackgroundWorkerHandle *handle; + + /* DecodingWorkerShared is in this segment. */ + dsm_segment *seg; + + /* Handle of the error queue. */ + shm_mq_handle *error_mqh; +} DecodingWorker; + +/* Pointer to currently running decoding worker. */ +static DecodingWorker *decoding_worker = NULL; + +/* + * Is there a message sent by a repack worker that the backend needs to + * receive? + */ +volatile sig_atomic_t RepackMessagePending = false; + static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -145,7 +262,7 @@ static void check_repack_concurrently_requirements(Relation rel); static void rebuild_relation(Relation OldHeap, Relation index, bool verbose, bool concurrent); static void copy_table_data(Relation NewHeap, Relation OldHeap, Relation OldIndex, - Snapshot snapshot, LogicalDecodingContext *decoding_ctx, + Snapshot snapshot, bool verbose, bool *pSwapToastByContent, TransactionId *pFreezeXid, @@ -158,12 +275,10 @@ static List *get_tables_to_repack_partitioned(RepackCommand cmd, static bool cluster_is_permitted_for_relation(RepackCommand cmd, Oid relid, Oid userid); -static void begin_concurrent_repack(Relation rel); -static void end_concurrent_repack(void); static LogicalDecodingContext *setup_logical_decoding(Oid relid); -static HeapTuple get_changed_tuple(char *change); -static void apply_concurrent_changes(RepackDecodingState *dstate, - ChangeDest *dest); +static bool decode_concurrent_changes(LogicalDecodingContext *ctx, + DecodingWorkerShared *shared); +static void apply_concurrent_changes(BufFile *file, ChangeDest *dest); static void apply_concurrent_insert(Relation rel, HeapTuple tup, IndexInsertState *iistate, TupleTableSlot *index_slot); @@ -175,9 +290,9 @@ static void apply_concurrent_delete(Relation rel, HeapTuple tup_target); static HeapTuple find_target_tuple(Relation rel, ChangeDest *dest, HeapTuple tup_key, TupleTableSlot *ident_slot); -static void process_concurrent_changes(LogicalDecodingContext *decoding_ctx, - XLogRecPtr end_of_wal, - ChangeDest *dest); +static void process_concurrent_changes(XLogRecPtr end_of_wal, + ChangeDest *dest, + bool done); static IndexInsertState *get_index_insert_state(Relation relation, Oid ident_index_id, Relation *ident_index_p); @@ -186,7 +301,6 @@ static ScanKey build_identity_key(Oid ident_idx_oid, Relation rel_src, static void free_index_insert_state(IndexInsertState *iistate); static void cleanup_logical_decoding(LogicalDecodingContext *ctx); static void rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, - LogicalDecodingContext *decoding_ctx, TransactionId frozenXid, MultiXactId cutoffMulti); static List *build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes); @@ -196,6 +310,13 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); +static void start_decoding_worker(Oid relid); +static void stop_decoding_worker(void); +static void repack_worker_internal(dsm_segment *seg); +static void export_initial_snapshot(Snapshot snapshot, + DecodingWorkerShared *shared); +static Snapshot get_initial_snapshot(DecodingWorker *worker); +static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -619,20 +740,20 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, /* rebuild_relation does all the dirty work */ PG_TRY(); { - /* - * For concurrent processing, make sure that our logical decoding - * ignores data changes of other tables than the one we are - * processing. - */ - if (concurrent) - begin_concurrent_repack(OldHeap); - rebuild_relation(OldHeap, index, verbose, concurrent); } PG_FINALLY(); { if (concurrent) - end_concurrent_repack(); + { + /* + * Since during normal operation the worker was already asked to + * exit, stopping it explicitly is especially important on ERROR. + * However it still seems a good practice to make sure that the + * worker never survives the REPACK command. + */ + stop_decoding_worker(); + } } PG_END_TRY(); @@ -929,7 +1050,6 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, bool concurrent bool swap_toast_by_content; TransactionId frozenXid; MultiXactId cutoffMulti; - LogicalDecodingContext *decoding_ctx = NULL; Snapshot snapshot = NULL; #if USE_ASSERT_CHECKING LOCKMODE lmode; @@ -943,19 +1063,36 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, bool concurrent if (concurrent) { /* - * Prepare to capture the concurrent data changes. + * The worker needs to be member of the locking group we're the leader + * of. We ought to become the leader before the worker starts. The + * worker will join the group as soon as it starts. * - * Note that this call waits for all transactions with XID already - * assigned to finish. If some of those transactions is waiting for a - * lock conflicting with ShareUpdateExclusiveLock on our table (e.g. - * it runs CREATE INDEX), we can end up in a deadlock. Not sure this - * risk is worth unlocking/locking the table (and its clustering - * index) and checking again if its still eligible for REPACK - * CONCURRENTLY. + * This is to make sure that the deadlock described below is + * detectable by deadlock.c: if the worker waits for a transaction to + * complete and we are waiting for the worker output, then effectively + * we (i.e. this backend) are waiting for that transaction. */ - decoding_ctx = setup_logical_decoding(tableOid); + BecomeLockGroupLeader(); + + /* + * Start the worker that decodes data changes applied while we're + * copying the table contents. + * + * Note that the worker has to wait for all transactions with XID + * already assigned to finish. If some of those transactions is + * waiting for a lock conflicting with ShareUpdateExclusiveLock on our + * table (e.g. it runs CREATE INDEX), we can end up in a deadlock. + * Not sure this risk is worth unlocking/locking the table (and its + * clustering index) and checking again if its still eligible for + * REPACK CONCURRENTLY. + */ + start_decoding_worker(tableOid); + + /* + * Wait until the worker has the initial snapshot and retrieve it. + */ + snapshot = get_initial_snapshot(decoding_worker); - snapshot = SnapBuildInitialSnapshotForRepack(decoding_ctx->snapshot_builder); PushActiveSnapshot(snapshot); } @@ -980,7 +1117,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, bool concurrent NewHeap = table_open(OIDNewHeap, NoLock); /* Copy the heap data into the new table in the desired order */ - copy_table_data(NewHeap, OldHeap, index, snapshot, decoding_ctx, verbose, + copy_table_data(NewHeap, OldHeap, index, snapshot, verbose, &swap_toast_by_content, &frozenXid, &cutoffMulti); /* The historic snapshot won't be needed anymore. */ @@ -1001,14 +1138,11 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, bool concurrent if (index) index_close(index, NoLock); - rebuild_relation_finish_concurrent(NewHeap, OldHeap, decoding_ctx, - frozenXid, cutoffMulti); + rebuild_relation_finish_concurrent(NewHeap, OldHeap, frozenXid, + cutoffMulti); pgstat_progress_update_param(PROGRESS_REPACK_PHASE, PROGRESS_REPACK_PHASE_FINAL_CLEANUP); - - /* Done with decoding. */ - cleanup_logical_decoding(decoding_ctx); } else { @@ -1179,8 +1313,7 @@ make_new_heap(Oid OIDOldHeap, Oid NewTableSpace, Oid NewAccessMethod, */ static void copy_table_data(Relation NewHeap, Relation OldHeap, Relation OldIndex, - Snapshot snapshot, LogicalDecodingContext *decoding_ctx, - bool verbose, bool *pSwapToastByContent, + Snapshot snapshot, bool verbose, bool *pSwapToastByContent, TransactionId *pFreezeXid, MultiXactId *pCutoffMulti) { Relation relRelation; @@ -1341,7 +1474,6 @@ copy_table_data(Relation NewHeap, Relation OldHeap, Relation OldIndex, */ table_relation_copy_for_cluster(OldHeap, NewHeap, OldIndex, use_sort, cutoffs.OldestXmin, snapshot, - decoding_ctx, &cutoffs.FreezeLimit, &cutoffs.MultiXactCutoff, &num_tuples, &tups_vacuumed, @@ -2375,58 +2507,6 @@ RepackCommandAsString(RepackCommand cmd) } -/* - * Call this function before REPACK CONCURRENTLY starts to setup logical - * decoding. It makes sure that other users of the table put enough - * information into WAL. - * - * The point is that at various places we expect that the table we're - * processing is treated like a system catalog. For example, we need to be - * able to scan it using a "historic snapshot" anytime during the processing - * (as opposed to scanning only at the start point of the decoding, as logical - * replication does during initial table synchronization), in order to apply - * concurrent UPDATE / DELETE commands. - * - * Note that TOAST table needs no attention here as it's not scanned using - * historic snapshot. - */ -static void -begin_concurrent_repack(Relation rel) -{ - Oid toastrelid; - - /* - * Avoid logical decoding of other relations by this backend. The lock we - * have guarantees that the actual locator cannot be changed concurrently: - * TRUNCATE needs AccessExclusiveLock. - */ - Assert(CheckRelationLockedByMe(rel, ShareUpdateExclusiveLock, false)); - repacked_rel_locator = rel->rd_locator; - toastrelid = rel->rd_rel->reltoastrelid; - if (OidIsValid(toastrelid)) - { - Relation toastrel; - - /* Avoid logical decoding of other TOAST relations. */ - toastrel = table_open(toastrelid, AccessShareLock); - repacked_rel_toast_locator = toastrel->rd_locator; - table_close(toastrel, AccessShareLock); - } -} - -/* - * Call this when done with REPACK CONCURRENTLY. - */ -static void -end_concurrent_repack(void) -{ - /* - * Restore normal function of (future) logical decoding for this backend. - */ - repacked_rel_locator.relNumber = InvalidOid; - repacked_rel_toast_locator.relNumber = InvalidOid; -} - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -2491,9 +2571,10 @@ static LogicalDecodingContext * setup_logical_decoding(Oid relid) { Relation rel; - TupleDesc tupdesc; + Oid toastrelid; LogicalDecodingContext *ctx; - RepackDecodingState *dstate = palloc0_object(RepackDecodingState); + NameData slotname; + RepackDecodingState *dstate; /* * REPACK CONCURRENTLY is not allowed in a transaction block, so this @@ -2501,21 +2582,21 @@ setup_logical_decoding(Oid relid) */ Assert(!TransactionIdIsValid(GetTopTransactionIdIfAny())); - /* - * A single backend should not execute multiple REPACK commands at a time, - * so use PID to make the slot unique. - */ - snprintf(NameStr(dstate->slotname), NAMEDATALEN, "repack_%d", MyProcPid); - /* * Check if we can use logical decoding. */ CheckSlotPermissions(); CheckLogicalDecodingRequirements(); - /* RS_TEMPORARY so that the slot gets cleaned up on ERROR. */ - ReplicationSlotCreate(NameStr(dstate->slotname), true, RS_TEMPORARY, - false, false, false); + /* + * A single backend should not execute multiple REPACK commands at a time, + * so use PID to make the slot unique. + * + * RS_TEMPORARY so that the slot gets cleaned up on ERROR. + */ + snprintf(NameStr(slotname), NAMEDATALEN, "repack_%d", MyProcPid); + ReplicationSlotCreate(NameStr(slotname), true, RS_TEMPORARY, false, false, + false); /* * Neither prepare_write nor do_write callback nor update_progress is @@ -2537,104 +2618,109 @@ setup_logical_decoding(Oid relid) DecodingContextFindStartpoint(ctx); + /* + * decode_concurrent_changes() needs non-blocking callback. + */ + ctx->reader->routine.page_read = read_local_xlog_page_no_wait; + + /* + * read_local_xlog_page_no_wait() needs to be able to indicate the end of + * WAL. + */ + ctx->reader->private_data = MemoryContextAllocZero(ctx->context, + sizeof(ReadLocalXLogPageNoWaitPrivate)); + + /* Some WAL records should have been read. */ Assert(ctx->reader->EndRecPtr != InvalidXLogRecPtr); + /* + * Initialize repack_current_segment so that we can notice WAL segment + * boundaries. + */ XLByteToSeg(ctx->reader->EndRecPtr, repack_current_segment, wal_segment_size); - /* - * Setup structures to store decoded changes. - */ + dstate = palloc0_object(RepackDecodingState); dstate->relid = relid; - dstate->tstore = tuplestore_begin_heap(false, false, - maintenance_work_mem); - /* Caller should already have the table locked. */ - rel = table_open(relid, NoLock); - tupdesc = CreateTupleDescCopy(RelationGetDescr(rel)); - dstate->tupdesc = tupdesc; - table_close(rel, NoLock); + /* + * Tuple descriptor may be needed to flatten a tuple before we write it to + * a file. A copy is needed because the decoding worker invalidates system + * caches before it starts to do the actual work. + */ + rel = table_open(relid, AccessShareLock); + dstate->tupdesc = CreateTupleDescCopy(RelationGetDescr(rel)); - /* Initialize the descriptor to store the changes ... */ - dstate->tupdesc_change = CreateTemplateTupleDesc(1); + /* Avoid logical decoding of other relations. */ + repacked_rel_locator = rel->rd_locator; + toastrelid = rel->rd_rel->reltoastrelid; + if (OidIsValid(toastrelid)) + { + Relation toastrel; - TupleDescInitEntry(dstate->tupdesc_change, 1, NULL, BYTEAOID, -1, 0); - /* ... as well as the corresponding slot. */ - dstate->tsslot = MakeSingleTupleTableSlot(dstate->tupdesc_change, - &TTSOpsMinimalTuple); + /* Avoid logical decoding of other TOAST relations. */ + toastrel = table_open(toastrelid, AccessShareLock); + repacked_rel_toast_locator = toastrel->rd_locator; + table_close(toastrel, AccessShareLock); + } + table_close(rel, AccessShareLock); - dstate->resowner = ResourceOwnerCreate(CurrentResourceOwner, - "logical decoding"); + /* The file will be set as soon as we have it opened. */ + dstate->file = NULL; ctx->output_writer_private = dstate; + return ctx; } /* - * Retrieve tuple from ConcurrentChange structure. + * Decode logical changes from the WAL sequence and store them to a file. * - * The input data starts with the structure but it might not be appropriately - * aligned. - */ -static HeapTuple -get_changed_tuple(char *change) -{ - HeapTupleData tup_data; - HeapTuple result; - char *src; - - /* - * Ensure alignment before accessing the fields. (This is why we can't use - * heap_copytuple() instead of this function.) - */ - src = change + offsetof(ConcurrentChange, tup_data); - memcpy(&tup_data, src, sizeof(HeapTupleData)); - - result = (HeapTuple) palloc(HEAPTUPLESIZE + tup_data.t_len); - memcpy(result, &tup_data, sizeof(HeapTupleData)); - result->t_data = (HeapTupleHeader) ((char *) result + HEAPTUPLESIZE); - src = change + SizeOfConcurrentChange; - memcpy(result->t_data, src, result->t_len); - - return result; -} - -/* - * Decode logical changes from the WAL sequence up to end_of_wal. + * If true is returned, there is no more work for the worker. */ -void -repack_decode_concurrent_changes(LogicalDecodingContext *ctx, - XLogRecPtr end_of_wal) +static bool +decode_concurrent_changes(LogicalDecodingContext *ctx, + DecodingWorkerShared *shared) { RepackDecodingState *dstate; - ResourceOwner resowner_old; + XLogRecPtr lsn_upto; + bool done; + char fname[MAXPGPATH]; dstate = (RepackDecodingState *) ctx->output_writer_private; - resowner_old = CurrentResourceOwner; - CurrentResourceOwner = dstate->resowner; - PG_TRY(); + /* Open the output file. */ + DecodingWorkerFileName(fname, shared->relid, shared->last_exported + 1); + dstate->file = BufFileCreateFileSet(&shared->sfs.fs, fname); + + SpinLockAcquire(&shared->mutex); + lsn_upto = shared->lsn_upto; + done = shared->done; + SpinLockRelease(&shared->mutex); + + while (true) { - while (ctx->reader->EndRecPtr < end_of_wal) - { - XLogRecord *record; - XLogSegNo segno_new; - char *errm = NULL; - XLogRecPtr end_lsn; + XLogRecord *record; + XLogSegNo segno_new; + char *errm = NULL; + XLogRecPtr end_lsn; - record = XLogReadRecord(ctx->reader, &errm); - if (errm) - elog(ERROR, "%s", errm); + CHECK_FOR_INTERRUPTS(); - if (record != NULL) - LogicalDecodingProcessRecord(ctx, ctx->reader); + record = XLogReadRecord(ctx->reader, &errm); + if (record) + { + LogicalDecodingProcessRecord(ctx, ctx->reader); /* * If WAL segment boundary has been crossed, inform the decoding - * system that the catalog_xmin can advance. (We can confirm more - * often, but a filling a single WAL segment should not take much - * time.) + * system that the catalog_xmin can advance. + * + * TODO Does it make sense to confirm more often? Segment size + * seems appropriate for restart_lsn (because less than a segment + * cannot be recycled anyway), however more frequent checks might + * be beneficial for catalog_xmin. */ end_lsn = ctx->reader->EndRecPtr; XLByteToSeg(end_lsn, segno_new, wal_segment_size); @@ -2645,80 +2731,137 @@ repack_decode_concurrent_changes(LogicalDecodingContext *ctx, (uint32) (end_lsn >> 32), (uint32) end_lsn); repack_current_segment = segno_new; } + } + else + { + ReadLocalXLogPageNoWaitPrivate *priv; - CHECK_FOR_INTERRUPTS(); + if (errm) + ereport(ERROR, (errmsg("%s", errm))); + + /* + * In the decoding loop we do not want to get blocked when there + * is no more WAL available, otherwise the loop would become + * uninterruptible. + */ + priv = (ReadLocalXLogPageNoWaitPrivate *) + ctx->reader->private_data; + if (priv->end_of_wal) + /* Do not miss the end of WAL condition next time. */ + priv->end_of_wal = false; + else + ereport(ERROR, (errmsg("could not read WAL record"))); + } + + /* + * Whether we could read new record or not, keep checking if + * 'lsn_upto' was specified. + */ + if (XLogRecPtrIsInvalid(lsn_upto)) + { + SpinLockAcquire(&shared->mutex); + lsn_upto = shared->lsn_upto; + /* 'done' should be set at the same time as 'lsn_upto' */ + done = shared->done; + SpinLockRelease(&shared->mutex); + } + if (!XLogRecPtrIsInvalid(lsn_upto) && + ctx->reader->EndRecPtr >= lsn_upto) + break; + + if (record == NULL) + { + int64 timeout = 0; + WaitLSNResult res; + + /* + * Before we retry reading, wait until new WAL is flushed. + * + * There is a race condition such that the backend executing + * REPACK determines 'lsn_upto', but before it sets the shared + * variable, we reach the end of WAL. In that case we'd need to + * wait until the next WAL flush (unrelated to REPACK). Although + * that should not be a problem in a busy system, it might be + * noticeable in other cases, including regression tests (which + * are not necessarily executed in parallel). Therefore it makes + * sense to use timeout. + * + * If lsn_upto is valid, WAL records having LSN lower than that + * should already have been flushed to disk. + */ + if (XLogRecPtrIsInvalid(lsn_upto)) + timeout = 100L; + res = WaitForLSN(WAIT_LSN_TYPE_PRIMARY_FLUSH, + ctx->reader->EndRecPtr + 1, + timeout); + if (res != WAIT_LSN_RESULT_SUCCESS && + res != WAIT_LSN_RESULT_TIMEOUT) + ereport(ERROR, (errmsg("waiting for WAL failed"))); } - InvalidateSystemCaches(); - CurrentResourceOwner = resowner_old; - } - PG_CATCH(); - { - /* clear all timetravel entries */ - InvalidateSystemCaches(); - CurrentResourceOwner = resowner_old; - PG_RE_THROW(); } - PG_END_TRY(); + + /* + * Close the file so we can make it available to the backend. + */ + BufFileClose(dstate->file); + dstate->file = NULL; + SpinLockAcquire(&shared->mutex); + shared->lsn_upto = InvalidXLogRecPtr; + shared->last_exported++; + SpinLockRelease(&shared->mutex); + ConditionVariableSignal(&shared->cv); + + return done; } /* * Apply changes stored in 'file'. */ static void -apply_concurrent_changes(RepackDecodingState *dstate, ChangeDest *dest) +apply_concurrent_changes(BufFile *file, ChangeDest *dest) { + char kind; + uint32 t_len; Relation rel = dest->rel; TupleTableSlot *index_slot, *ident_slot; HeapTuple tup_old = NULL; - if (dstate->nchanges == 0) - return; - /* TupleTableSlot is needed to pass the tuple to ExecInsertIndexTuples(). */ - index_slot = MakeSingleTupleTableSlot(dstate->tupdesc, &TTSOpsHeapTuple); + index_slot = MakeSingleTupleTableSlot(RelationGetDescr(rel), + &TTSOpsHeapTuple); /* A slot to fetch tuples from identity index. */ ident_slot = table_slot_create(rel, NULL); - while (tuplestore_gettupleslot(dstate->tstore, true, false, - dstate->tsslot)) + while (true) { - bool shouldFree; - HeapTuple tup_change, - tup, + size_t nread; + HeapTuple tup, tup_exist; - char *change_raw, - *src; - ConcurrentChange change; - bool isnull[1]; - Datum values[1]; CHECK_FOR_INTERRUPTS(); - /* Get the change from the single-column tuple. */ - tup_change = ExecFetchSlotHeapTuple(dstate->tsslot, false, &shouldFree); - heap_deform_tuple(tup_change, dstate->tupdesc_change, values, isnull); - Assert(!isnull[0]); - - /* Make sure we access aligned data. */ - change_raw = (char *) DatumGetByteaP(values[0]); - src = (char *) VARDATA(change_raw); - memcpy(&change, src, SizeOfConcurrentChange); + nread = BufFileReadMaybeEOF(file, &kind, 1, true); + /* Are we done with the file? */ + if (nread == 0) + break; - /* - * Extract the tuple from the change. The tuple is copied here because - * it might be assigned to 'tup_old', in which case it needs to - * survive into the next iteration. - */ - tup = get_changed_tuple(src); + /* Read the tuple. */ + BufFileReadExact(file, &t_len, sizeof(t_len)); + tup = (HeapTuple) palloc(HEAPTUPLESIZE + t_len); + tup->t_data = (HeapTupleHeader) ((char *) tup + HEAPTUPLESIZE); + BufFileReadExact(file, tup->t_data, t_len); + tup->t_len = t_len; + ItemPointerSetInvalid(&tup->t_self); + tup->t_tableOid = RelationGetRelid(dest->rel); - if (change.kind == CHANGE_UPDATE_OLD) + if (kind == CHANGE_UPDATE_OLD) { Assert(tup_old == NULL); tup_old = tup; } - else if (change.kind == CHANGE_INSERT) + else if (kind == CHANGE_INSERT) { Assert(tup_old == NULL); @@ -2726,12 +2869,11 @@ apply_concurrent_changes(RepackDecodingState *dstate, ChangeDest *dest) pfree(tup); } - else if (change.kind == CHANGE_UPDATE_NEW || - change.kind == CHANGE_DELETE) + else if (kind == CHANGE_UPDATE_NEW || kind == CHANGE_DELETE) { HeapTuple tup_key; - if (change.kind == CHANGE_UPDATE_NEW) + if (kind == CHANGE_UPDATE_NEW) { tup_key = tup_old != NULL ? tup_old : tup; } @@ -2748,7 +2890,7 @@ apply_concurrent_changes(RepackDecodingState *dstate, ChangeDest *dest) if (tup_exist == NULL) elog(ERROR, "failed to find target tuple"); - if (change.kind == CHANGE_UPDATE_NEW) + if (kind == CHANGE_UPDATE_NEW) apply_concurrent_update(rel, tup, tup_exist, dest->iistate, index_slot); else @@ -2763,26 +2905,19 @@ apply_concurrent_changes(RepackDecodingState *dstate, ChangeDest *dest) pfree(tup); } else - elog(ERROR, "unrecognized kind of change: %d", change.kind); + elog(ERROR, "unrecognized kind of change: %d", kind); /* * If a change was applied now, increment CID for next writes and * update the snapshot so it sees the changes we've applied so far. */ - if (change.kind != CHANGE_UPDATE_OLD) + if (kind != CHANGE_UPDATE_OLD) { CommandCounterIncrement(); UpdateActiveSnapshotCommandId(); } - - /* TTSOpsMinimalTuple has .get_heap_tuple==NULL. */ - Assert(shouldFree); - pfree(tup_change); } - tuplestore_clear(dstate->tstore); - dstate->nchanges = 0; - /* Cleanup. */ ExecDropSingleTupleTableSlot(index_slot); ExecDropSingleTupleTableSlot(ident_slot); @@ -2957,25 +3092,59 @@ find_target_tuple(Relation rel, ChangeDest *dest, HeapTuple tup_key, } /* - * Decode and apply concurrent changes. + * Decode and apply concurrent changes, up to (and including) the record whose + * LSN is 'end_of_wal'. */ static void -process_concurrent_changes(LogicalDecodingContext *decoding_ctx, - XLogRecPtr end_of_wal, ChangeDest *dest) +process_concurrent_changes(XLogRecPtr end_of_wal, ChangeDest *dest, bool done) { - RepackDecodingState *dstate; + DecodingWorkerShared *shared; + char fname[MAXPGPATH]; + BufFile *file; pgstat_progress_update_param(PROGRESS_REPACK_PHASE, PROGRESS_REPACK_PHASE_CATCH_UP); - dstate = (RepackDecodingState *) decoding_ctx->output_writer_private; + /* Ask the worker for the file. */ + shared = (DecodingWorkerShared *) dsm_segment_address(decoding_worker->seg); + SpinLockAcquire(&shared->mutex); + shared->lsn_upto = end_of_wal; + shared->done = done; + SpinLockRelease(&shared->mutex); + + /* + * The worker needs to finish processing of the current WAL record. Even + * if it's idle, it'll need to close the output file. Thus we're likely to + * wait, so prepare for sleep. + */ + ConditionVariablePrepareToSleep(&shared->cv); + for (;;) + { + int last_exported; - repack_decode_concurrent_changes(decoding_ctx, end_of_wal); + SpinLockAcquire(&shared->mutex); + last_exported = shared->last_exported; + SpinLockRelease(&shared->mutex); - if (dstate->nchanges == 0) - return; + /* + * Has the worker exported the file we are waiting for? + */ + if (last_exported == dest->file_seq) + break; - apply_concurrent_changes(dstate, dest); + ConditionVariableSleep(&shared->cv, WAIT_EVENT_REPACK_WORKER_EXPORT); + } + ConditionVariableCancelSleep(); + + /* Open the file. */ + DecodingWorkerFileName(fname, shared->relid, dest->file_seq); + file = BufFileOpenFileSet(&shared->sfs.fs, fname, O_RDONLY, false); + apply_concurrent_changes(file, dest); + + BufFileClose(file); + + /* Get ready for the next file. */ + dest->file_seq++; } /* @@ -3101,15 +3270,10 @@ cleanup_logical_decoding(LogicalDecodingContext *ctx) dstate = (RepackDecodingState *) ctx->output_writer_private; - ExecDropSingleTupleTableSlot(dstate->tsslot); - FreeTupleDesc(dstate->tupdesc_change); FreeTupleDesc(dstate->tupdesc); - tuplestore_end(dstate->tstore); - FreeDecodingContext(ctx); - ReplicationSlotRelease(); - ReplicationSlotDrop(NameStr(dstate->slotname), false); + ReplicationSlotDropAcquired(); pfree(dstate); } @@ -3123,7 +3287,6 @@ cleanup_logical_decoding(LogicalDecodingContext *ctx) */ static void rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, - LogicalDecodingContext *decoding_ctx, TransactionId frozenXid, MultiXactId cutoffMulti) { @@ -3204,6 +3367,7 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, &chgdst.ident_index); chgdst.ident_key = build_identity_key(ident_idx_new, OldHeap, &chgdst.ident_key_nentries); + chgdst.file_seq = WORKER_FILE_SNAPSHOT + 1; /* * During testing, wait for another backend to perform concurrent data @@ -3225,7 +3389,7 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, * hold AccessExclusiveLock. (Quite some amount of WAL could have been * written during the data copying and index creation.) */ - process_concurrent_changes(decoding_ctx, end_of_wal, &chgdst); + process_concurrent_changes(end_of_wal, &chgdst, false); /* * Acquire AccessExclusiveLock on the table, its TOAST relation (if there @@ -3306,8 +3470,11 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, XLogFlush(wal_insert_ptr); end_of_wal = GetFlushRecPtr(NULL); - /* Apply the concurrent changes again. */ - process_concurrent_changes(decoding_ctx, end_of_wal, &chgdst); + /* + * Apply the concurrent changes again. Indicate that the decoding worker + * won't be needed anymore. + */ + process_concurrent_changes(end_of_wal, &chgdst, true); /* Remember info about rel before closing OldHeap */ relpersistence = OldHeap->rd_rel->relpersistence; @@ -3417,3 +3584,510 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) return result; } + +/* + * Try to start a background worker to perform logical decoding of data + * changes applied to relation while REPACK CONCURRENTLY is copying its + * contents to a new table. + */ +static void +start_decoding_worker(Oid relid) +{ + Size size; + dsm_segment *seg; + DecodingWorkerShared *shared; + shm_mq *mq; + shm_mq_handle *mqh; + BackgroundWorker bgw; + + /* Setup shared memory. */ + size = BUFFERALIGN(offsetof(DecodingWorkerShared, error_queue)) + + BUFFERALIGN(REPACK_ERROR_QUEUE_SIZE); + seg = dsm_create(size, 0); + shared = (DecodingWorkerShared *) dsm_segment_address(seg); + shared->lsn_upto = InvalidXLogRecPtr; + shared->done = false; + SharedFileSetInit(&shared->sfs, seg); + shared->last_exported = -1; + SpinLockInit(&shared->mutex); + shared->dbid = MyDatabaseId; + + /* + * This is the UserId set in cluster_rel(). Security context shouldn't be + * needed for decoding worker. + */ + shared->roleid = GetUserId(); + shared->relid = relid; + ConditionVariableInit(&shared->cv); + shared->backend_proc = MyProc; + shared->backend_pid = MyProcPid; + shared->backend_proc_number = MyProcNumber; + + mq = shm_mq_create((char *) BUFFERALIGN(shared->error_queue), + REPACK_ERROR_QUEUE_SIZE); + shm_mq_set_receiver(mq, MyProc); + mqh = shm_mq_attach(mq, seg, NULL); + + memset(&bgw, 0, sizeof(bgw)); + snprintf(bgw.bgw_name, BGW_MAXLEN, + "REPACK decoding worker for relation \"%s\"", + get_rel_name(relid)); + snprintf(bgw.bgw_type, BGW_MAXLEN, "REPACK decoding worker"); + bgw.bgw_flags = BGWORKER_SHMEM_ACCESS | + BGWORKER_BACKEND_DATABASE_CONNECTION; + bgw.bgw_start_time = BgWorkerStart_RecoveryFinished; + bgw.bgw_restart_time = BGW_NEVER_RESTART; + snprintf(bgw.bgw_library_name, MAXPGPATH, "postgres"); + snprintf(bgw.bgw_function_name, BGW_MAXLEN, "RepackWorkerMain"); + bgw.bgw_main_arg = UInt32GetDatum(dsm_segment_handle(seg)); + bgw.bgw_notify_pid = MyProcPid; + + decoding_worker = palloc0_object(DecodingWorker); + if (!RegisterDynamicBackgroundWorker(&bgw, &decoding_worker->handle)) + ereport(ERROR, + (errcode(ERRCODE_CONFIGURATION_LIMIT_EXCEEDED), + errmsg("out of background worker slots"), + errhint("You might need to increase \"%s\".", "max_worker_processes"))); + + decoding_worker->seg = seg; + decoding_worker->error_mqh = mqh; + + /* + * The decoding setup must be done before the caller can have XID assigned + * for any reason, otherwise the worker might end up in a deadlock, + * waiting for the caller's transaction to end. Therefore wait here until + * the worker indicates that it has the logical decoding initialized. + */ + ConditionVariablePrepareToSleep(&shared->cv); + for (;;) + { + int initialized; + + SpinLockAcquire(&shared->mutex); + initialized = shared->initialized; + SpinLockRelease(&shared->mutex); + + if (initialized) + break; + + ConditionVariableSleep(&shared->cv, WAIT_EVENT_REPACK_WORKER_EXPORT); + } + ConditionVariableCancelSleep(); +} + +/* + * Stop the decoding worker and cleanup the related resources. + * + * The worker stops on its own when it knows there is no more work to do, but + * we need to stop it explicitly at least on ERROR in the launching backend. + */ +static void +stop_decoding_worker(void) +{ + BgwHandleStatus status; + + /* Haven't reached the worker startup? */ + if (decoding_worker == NULL) + return; + + /* Could not register the worker? */ + if (decoding_worker->handle == NULL) + return; + + TerminateBackgroundWorker(decoding_worker->handle); + /* The worker should really exit before the REPACK command does. */ + HOLD_INTERRUPTS(); + status = WaitForBackgroundWorkerShutdown(decoding_worker->handle); + RESUME_INTERRUPTS(); + + if (status == BGWH_POSTMASTER_DIED) + ereport(FATAL, + (errcode(ERRCODE_ADMIN_SHUTDOWN), + errmsg("postmaster exited during REPACK command"))); + + shm_mq_detach(decoding_worker->error_mqh); + + /* + * If we could not cancel the current sleep due to ERROR, do that before + * we detach from the shared memory the condition variable is located in. + * If we did not, the bgworker ERROR handling code would try and fail + * badly. + */ + ConditionVariableCancelSleep(); + + dsm_detach(decoding_worker->seg); + pfree(decoding_worker); + decoding_worker = NULL; +} + +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; + +/* + * See ParallelWorkerShutdown for details. + */ +static void +RepackWorkerShutdown(int code, Datum arg) +{ + SendProcSignal(backend_pid, + PROCSIG_REPACK_MESSAGE, + backend_proc_number); + + dsm_detach((dsm_segment *) DatumGetPointer(arg)); +} + +/* REPACK decoding worker entry point */ +void +RepackWorkerMain(Datum main_arg) +{ + dsm_segment *seg; + DecodingWorkerShared *shared; + shm_mq *mq; + shm_mq_handle *mqh; + + is_repack_worker = true; + + /* + * Override the default bgworker_die() with die() so we can use + * CHECK_FOR_INTERRUPTS(). + */ + pqsignal(SIGTERM, die); + BackgroundWorkerUnblockSignals(); + + seg = dsm_attach(DatumGetUInt32(main_arg)); + if (seg == NULL) + ereport(ERROR, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("could not map dynamic shared memory segment"))); + + shared = (DecodingWorkerShared *) dsm_segment_address(seg); + + /* Arrange to signal the leader if we exit. */ + backend_pid = shared->backend_pid; + backend_proc_number = shared->backend_proc_number; + before_shmem_exit(RepackWorkerShutdown, PointerGetDatum(seg)); + + /* + * Join locking group - see the comments around the call of + * start_decoding_worker(). + */ + if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) + /* The leader is not running anymore. */ + return; + + /* + * Setup a queue to send error messages to the backend that launched this + * worker. + */ + mq = (shm_mq *) (char *) BUFFERALIGN(shared->error_queue); + shm_mq_set_sender(mq, MyProc); + mqh = shm_mq_attach(mq, seg, NULL); + pq_redirect_to_shm_mq(seg, mqh); + pq_set_parallel_leader(shared->backend_pid, + shared->backend_proc_number); + + /* Connect to the database. */ + BackgroundWorkerInitializeConnectionByOid(shared->dbid, shared->roleid, 0); + + repack_worker_internal(seg); +} + +static void +repack_worker_internal(dsm_segment *seg) +{ + DecodingWorkerShared *shared; + LogicalDecodingContext *decoding_ctx; + SharedFileSet *sfs; + Snapshot snapshot; + + /* + * Transaction is needed to open relation, and it also provides us with a + * resource owner. + */ + StartTransactionCommand(); + + shared = (DecodingWorkerShared *) dsm_segment_address(seg); + + /* + * Not sure the spinlock is needed here - the backend should not change + * anything in the shared memory until we have serialized the snapshot. + */ + SpinLockAcquire(&shared->mutex); + Assert(XLogRecPtrIsInvalid(shared->lsn_upto)); + sfs = &shared->sfs; + SpinLockRelease(&shared->mutex); + + SharedFileSetAttach(sfs, seg); + + /* + * Prepare to capture the concurrent data changes ourselves. + */ + decoding_ctx = setup_logical_decoding(shared->relid); + + /* Announce that we're ready. */ + SpinLockAcquire(&shared->mutex); + shared->initialized = true; + SpinLockRelease(&shared->mutex); + ConditionVariableSignal(&shared->cv); + + /* Build the initial snapshot and export it. */ + snapshot = SnapBuildInitialSnapshotForRepack(decoding_ctx->snapshot_builder); + export_initial_snapshot(snapshot, shared); + + /* + * Only historic snapshots should be used now. Do not let us restrict the + * progress of xmin horizon. + */ + InvalidateCatalogSnapshot(); + + while (!decode_concurrent_changes(decoding_ctx, shared)) + ; + + /* Cleanup. */ + cleanup_logical_decoding(decoding_ctx); + CommitTransactionCommand(); +} + +/* + * Make snapshot available to the backend that launched the decoding worker. + */ +static void +export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared) +{ + char fname[MAXPGPATH]; + BufFile *file; + Size snap_size; + char *snap_space; + + snap_size = EstimateSnapshotSpace(snapshot); + snap_space = (char *) palloc(snap_size); + SerializeSnapshot(snapshot, snap_space); + FreeSnapshot(snapshot); + + DecodingWorkerFileName(fname, shared->relid, shared->last_exported + 1); + file = BufFileCreateFileSet(&shared->sfs.fs, fname); + /* To make restoration easier, write the snapshot size first. */ + BufFileWrite(file, &snap_size, sizeof(snap_size)); + BufFileWrite(file, snap_space, snap_size); + pfree(snap_space); + BufFileClose(file); + + /* Increase the counter to tell the backend that the file is available. */ + SpinLockAcquire(&shared->mutex); + shared->last_exported++; + SpinLockRelease(&shared->mutex); + ConditionVariableSignal(&shared->cv); +} + +/* + * Get the initial snapshot from the decoding worker. + */ +static Snapshot +get_initial_snapshot(DecodingWorker *worker) +{ + DecodingWorkerShared *shared; + char fname[MAXPGPATH]; + BufFile *file; + Size snap_size; + char *snap_space; + Snapshot snapshot; + + shared = (DecodingWorkerShared *) dsm_segment_address(worker->seg); + + /* + * The worker needs to initialize the logical decoding, which usually + * takes some time. Therefore it makes sense to prepare for the sleep + * first. + */ + ConditionVariablePrepareToSleep(&shared->cv); + for (;;) + { + int last_exported; + + SpinLockAcquire(&shared->mutex); + last_exported = shared->last_exported; + SpinLockRelease(&shared->mutex); + + /* + * Has the worker exported the file we are waiting for? + */ + if (last_exported == WORKER_FILE_SNAPSHOT) + break; + + ConditionVariableSleep(&shared->cv, WAIT_EVENT_REPACK_WORKER_EXPORT); + } + ConditionVariableCancelSleep(); + + /* Read the snapshot from a file. */ + DecodingWorkerFileName(fname, shared->relid, WORKER_FILE_SNAPSHOT); + file = BufFileOpenFileSet(&shared->sfs.fs, fname, O_RDONLY, false); + BufFileReadExact(file, &snap_size, sizeof(snap_size)); + snap_space = (char *) palloc(snap_size); + BufFileReadExact(file, snap_space, snap_size); + BufFileClose(file); + + /* Restore it. */ + snapshot = RestoreSnapshot(snap_space); + pfree(snap_space); + + return snapshot; +} + +bool +IsRepackWorker(void) +{ + return is_repack_worker; +} + +/* + * Handle receipt of an interrupt indicating a repack worker message. + * + * Note: this is called within a signal handler! All we can do is set + * a flag that will cause the next CHECK_FOR_INTERRUPTS() to invoke + * ProcessRepackMessages(). + */ +void +HandleRepackMessageInterrupt(void) +{ + InterruptPending = true; + RepackMessagePending = true; + SetLatch(MyLatch); +} + +/* + * Process any queued protocol messages received from parallel workers. + */ +void +ProcessRepackMessages(void) +{ + MemoryContext oldcontext; + + static MemoryContext hpm_context = NULL; + + /* + * Nothing to do if we haven't launched the worker yet or have already + * terminated it. + */ + if (decoding_worker == NULL) + return; + + /* + * This is invoked from ProcessInterrupts(), and since some of the + * functions it calls contain CHECK_FOR_INTERRUPTS(), there is a potential + * for recursive calls if more signals are received while this runs. It's + * unclear that recursive entry would be safe, and it doesn't seem useful + * even if it is safe, so let's block interrupts until done. + */ + HOLD_INTERRUPTS(); + + /* + * Moreover, CurrentMemoryContext might be pointing almost anywhere. We + * don't want to risk leaking data into long-lived contexts, so let's do + * our work here in a private context that we can reset on each use. + */ + if (hpm_context == NULL) /* first time through? */ + hpm_context = AllocSetContextCreate(TopMemoryContext, + "ProcessRepackMessages", + ALLOCSET_DEFAULT_SIZES); + else + MemoryContextReset(hpm_context); + + oldcontext = MemoryContextSwitchTo(hpm_context); + + /* OK to process messages. Reset the flag saying there are more to do. */ + RepackMessagePending = false; + + /* + * Read as many messages as we can from each worker, but stop when no more + * messages can be read from the worker without blocking. + */ + while (true) + { + shm_mq_result res; + Size nbytes; + void *data; + + res = shm_mq_receive(decoding_worker->error_mqh, &nbytes, + &data, true); + if (res == SHM_MQ_WOULD_BLOCK) + break; + else if (res == SHM_MQ_SUCCESS) + { + StringInfoData msg; + + initStringInfo(&msg); + appendBinaryStringInfo(&msg, data, nbytes); + ProcessRepackMessage(&msg); + pfree(msg.data); + } + else + { + /* + * The decoding worker is special in that it exits as soon as it + * has its work done. Thus the DETACHED result code is fine. + */ + Assert(res == SHM_MQ_DETACHED); + + break; + } + } + + MemoryContextSwitchTo(oldcontext); + + /* Might as well clear the context on our way out */ + MemoryContextReset(hpm_context); + + RESUME_INTERRUPTS(); +} + +/* + * Process a single protocol message received from a single parallel worker. + */ +static void +ProcessRepackMessage(StringInfo msg) +{ + char msgtype; + + msgtype = pq_getmsgbyte(msg); + + switch (msgtype) + { + case PqMsg_ErrorResponse: + case PqMsg_NoticeResponse: + { + ErrorData edata; + + /* Parse ErrorResponse or NoticeResponse. */ + pq_parse_errornotice(msg, &edata); + + /* Death of a worker isn't enough justification for suicide. */ + edata.elevel = Min(edata.elevel, ERROR); + + /* + * If desired, add a context line to show that this is a + * message propagated from a parallel worker. Otherwise, it + * can sometimes be confusing to understand what actually + * happened. + */ + if (edata.context) + edata.context = psprintf("%s\n%s", edata.context, + _("decoding worker")); + else + edata.context = pstrdup(_("decoding worker")); + + /* Rethrow error or print notice. */ + ThrowErrorData(&edata); + + break; + } + + default: + { + elog(ERROR, "unrecognized message type received from decoding worker: %c (message length %d bytes)", + msgtype, msg->len); + } + } +} diff --git a/src/backend/libpq/pqmq.c b/src/backend/libpq/pqmq.c index 6e4bbfb5aa1..42f6fa472c5 100644 --- a/src/backend/libpq/pqmq.c +++ b/src/backend/libpq/pqmq.c @@ -14,6 +14,7 @@ #include "postgres.h" #include "access/parallel.h" +#include "commands/cluster.h" #include "libpq/libpq.h" #include "libpq/pqformat.h" #include "libpq/pqmq.h" @@ -175,6 +176,10 @@ mq_putmessage(char msgtype, const char *s, size_t len) SendProcSignal(pq_mq_parallel_leader_pid, PROCSIG_PARALLEL_APPLY_MESSAGE, pq_mq_parallel_leader_proc_number); + else if (IsRepackWorker()) + SendProcSignal(pq_mq_parallel_leader_pid, + PROCSIG_REPACK_MESSAGE, + pq_mq_parallel_leader_proc_number); else { Assert(IsParallelWorker()); diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index 65deabe91a7..334bb708c5b 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,6 +13,7 @@ #include "postgres.h" #include "access/parallel.h" +#include "commands/cluster.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" @@ -136,6 +137,9 @@ static const struct }, { "SequenceSyncWorkerMain", SequenceSyncWorkerMain + }, + { + "RepackWorkerMain", RepackWorkerMain } }; diff --git a/src/backend/replication/logical/logical.c b/src/backend/replication/logical/logical.c index b0ef1a12520..35a46988285 100644 --- a/src/backend/replication/logical/logical.c +++ b/src/backend/replication/logical/logical.c @@ -194,7 +194,11 @@ StartupDecodingContext(List *output_plugin_options, ctx->slot = slot; - ctx->reader = XLogReaderAllocate(wal_segment_size, NULL, xl_routine, ctx); + /* + * TODO A separate patch for PG core, unless there's really a reason to + * pass ctx for private_data (May extensions expect ctx?). + */ + ctx->reader = XLogReaderAllocate(wal_segment_size, NULL, xl_routine, NULL); if (!ctx->reader) ereport(ERROR, (errcode(ERRCODE_OUT_OF_MEMORY), diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index a074eed393d..4bc47a72371 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -169,17 +169,13 @@ store_change(LogicalDecodingContext *ctx, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - char *change_raw; - ConcurrentChange change; + char kind_byte = (char) kind; bool flattened = false; - Size size; - Datum values[1]; - bool isnull[1]; - char *dst; dstate = (RepackDecodingState *) ctx->output_writer_private; - size = VARHDRSZ + SizeOfConcurrentChange; + /* Store the change kind. */ + BufFileWrite(dstate->file, &kind_byte, 1); /* * ReorderBufferCommit() stores the TOAST chunks in its private memory @@ -196,46 +192,12 @@ store_change(LogicalDecodingContext *ctx, ConcurrentChangeKind kind, tuple = toast_flatten_tuple(tuple, dstate->tupdesc); flattened = true; } + /* Store the tuple size ... */ + BufFileWrite(dstate->file, &tuple->t_len, sizeof(tuple->t_len)); + /* ... and the tuple itself. */ + BufFileWrite(dstate->file, tuple->t_data, tuple->t_len); - size += tuple->t_len; - if (size >= MaxAllocSize) - elog(ERROR, "Change is too big."); - - /* Construct the change. */ - change_raw = (char *) palloc0(size); - SET_VARSIZE(change_raw, size); - - /* - * Since the varlena alignment might not be sufficient for the structure, - * set the fields in a local instance and remember where it should - * eventually be copied. - */ - change.kind = kind; - dst = (char *) VARDATA(change_raw); - - /* - * Copy the tuple. - * - * Note: change->tup_data.t_data must be fixed on retrieval! - */ - memcpy(&change.tup_data, tuple, sizeof(HeapTupleData)); - memcpy(dst, &change, SizeOfConcurrentChange); - dst += SizeOfConcurrentChange; - memcpy(dst, tuple->t_data, tuple->t_len); - - /* The data has been copied. */ + /* Free the flat copy if created above. */ if (flattened) pfree(tuple); - - /* Store as tuple of 1 bytea column. */ - values[0] = PointerGetDatum(change_raw); - isnull[0] = false; - tuplestore_putvalues(dstate->tstore, dstate->tupdesc_change, - values, isnull); - - /* Accounting. */ - dstate->nchanges++; - - /* Cleanup. */ - pfree(change_raw); } diff --git a/src/backend/storage/ipc/procsignal.c b/src/backend/storage/ipc/procsignal.c index 8e56922dcea..6f9e7a7aab7 100644 --- a/src/backend/storage/ipc/procsignal.c +++ b/src/backend/storage/ipc/procsignal.c @@ -19,6 +19,7 @@ #include "access/parallel.h" #include "commands/async.h" +#include "commands/cluster.h" #include "miscadmin.h" #include "pgstat.h" #include "port/pg_bitutils.h" @@ -697,6 +698,9 @@ procsignal_sigusr1_handler(SIGNAL_ARGS) if (CheckProcSignal(PROCSIG_PARALLEL_APPLY_MESSAGE)) HandleParallelApplyMessageInterrupt(); + if (CheckProcSignal(PROCSIG_REPACK_MESSAGE)) + HandleRepackMessageInterrupt(); + if (CheckProcSignal(PROCSIG_RECOVERY_CONFLICT_DATABASE)) HandleRecoveryConflictInterrupt(PROCSIG_RECOVERY_CONFLICT_DATABASE); diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c index e54bf1e760f..fc81ad87615 100644 --- a/src/backend/tcop/postgres.c +++ b/src/backend/tcop/postgres.c @@ -36,6 +36,7 @@ #include "access/xact.h" #include "catalog/pg_type.h" #include "commands/async.h" +#include "commands/cluster.h" #include "commands/event_trigger.h" #include "commands/explain_state.h" #include "commands/prepare.h" @@ -3541,6 +3542,9 @@ ProcessInterrupts(void) if (ParallelApplyMessagePending) ProcessParallelApplyMessages(); + + if (RepackMessagePending) + ProcessRepackMessages(); } /* diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 3299de23bb3..73a3def69bc 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -62,6 +62,7 @@ LOGICAL_APPLY_MAIN "Waiting in main loop of logical replication apply process." LOGICAL_LAUNCHER_MAIN "Waiting in main loop of logical replication launcher process." LOGICAL_PARALLEL_APPLY_MAIN "Waiting in main loop of logical replication parallel apply process." RECOVERY_WAL_STREAM "Waiting in main loop of startup process for WAL to arrive, during streaming recovery." +REPACK_WORKER_MAIN "Waiting in main loop of REPACK decoding worker process." REPLICATION_SLOTSYNC_MAIN "Waiting in main loop of slot synchronization." REPLICATION_SLOTSYNC_SHUTDOWN "Waiting for slot sync worker to shut down." SYSLOGGER_MAIN "Waiting in main loop of syslogger process." @@ -154,6 +155,7 @@ RECOVERY_CONFLICT_SNAPSHOT "Waiting for recovery conflict resolution for a vacuu RECOVERY_CONFLICT_TABLESPACE "Waiting for recovery conflict resolution for dropping a tablespace." RECOVERY_END_COMMAND "Waiting for <xref linkend="guc-recovery-end-command"/> to complete." RECOVERY_PAUSE "Waiting for recovery to be resumed." +REPACK_WORKER_EXPORT "Waiting for decoding worker to export a new output file." REPLICATION_ORIGIN_DROP "Waiting for a replication origin to become inactive so it can be dropped." REPLICATION_SLOT_DROP "Waiting for a replication slot to become inactive so it can be dropped." RESTORE_COMMAND "Waiting for <xref linkend="guc-restore-command"/> to complete." diff --git a/src/include/access/tableam.h b/src/include/access/tableam.h index 76aa993009a..15760363a1a 100644 --- a/src/include/access/tableam.h +++ b/src/include/access/tableam.h @@ -22,7 +22,6 @@ #include "access/xact.h" #include "commands/vacuum.h" #include "executor/tuptable.h" -#include "replication/logical.h" #include "storage/read_stream.h" #include "utils/rel.h" #include "utils/snapshot.h" @@ -631,7 +630,6 @@ typedef struct TableAmRoutine bool use_sort, TransactionId OldestXmin, Snapshot snapshot, - LogicalDecodingContext *decoding_ctx, TransactionId *xid_cutoff, MultiXactId *multi_cutoff, double *num_tuples, @@ -1651,8 +1649,6 @@ table_relation_copy_data(Relation rel, const RelFileLocator *newrlocator) * - *multi_cutoff - ditto * - snapshot - if != NULL, ignore data changes done by transactions that this * (MVCC) snapshot considers still in-progress or in the future. - * - decoding_ctx - logical decoding context, to capture concurrent data - * changes. * * Output parameters: * - *xid_cutoff - rel's new relfrozenxid value, may be invalid @@ -1666,7 +1662,6 @@ table_relation_copy_for_cluster(Relation OldTable, Relation NewTable, bool use_sort, TransactionId OldestXmin, Snapshot snapshot, - LogicalDecodingContext *decoding_ctx, TransactionId *xid_cutoff, MultiXactId *multi_cutoff, double *num_tuples, @@ -1675,7 +1670,7 @@ table_relation_copy_for_cluster(Relation OldTable, Relation NewTable, { OldTable->rd_tableam->relation_copy_for_cluster(OldTable, NewTable, OldIndex, use_sort, OldestXmin, - snapshot, decoding_ctx, + snapshot, xid_cutoff, multi_cutoff, num_tuples, tups_vacuumed, tups_recently_dead); diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 6a5c476294a..1b05d5d418b 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -17,11 +17,13 @@ #include "nodes/parsenodes.h" #include "parser/parse_node.h" #include "replication/decode.h" +#include "postmaster/bgworker.h" #include "replication/logical.h" +#include "storage/buffile.h" #include "storage/lock.h" +#include "storage/shm_mq.h" #include "utils/relcache.h" #include "utils/resowner.h" -#include "utils/tuplestore.h" /* flag bits for ClusterParams->options */ @@ -44,6 +46,9 @@ typedef struct ClusterParams * The following definitions are used by REPACK CONCURRENTLY. */ +/* + * Stored as a single byte in the output file. + */ typedef enum { CHANGE_INSERT, @@ -52,68 +57,30 @@ typedef enum CHANGE_DELETE } ConcurrentChangeKind; -typedef struct ConcurrentChange -{ - /* See the enum above. */ - ConcurrentChangeKind kind; - - /* - * The actual tuple. - * - * The tuple data follows the ConcurrentChange structure. Before use make - * sure the tuple is correctly aligned (ConcurrentChange can be stored as - * bytea) and that tuple->t_data is fixed. - */ - HeapTupleData tup_data; -} ConcurrentChange; - -#define SizeOfConcurrentChange (offsetof(ConcurrentChange, tup_data) + \ - sizeof(HeapTupleData)) - /* * Logical decoding state. * - * Here we store the data changes that we decode from WAL while the table - * contents is being copied to a new storage. Also the necessary metadata - * needed to apply these changes to the table is stored here. + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. */ typedef struct RepackDecodingState { /* The relation whose changes we're decoding. */ Oid relid; - /* Replication slot name. */ - NameData slotname; - - /* - * Decoded changes are stored here. Although we try to avoid excessive - * batches, it can happen that the changes need to be stored to disk. The - * tuplestore does this transparently. - */ - Tuplestorestate *tstore; - - /* The current number of changes in tstore. */ - double nchanges; - - /* - * Descriptor to store the ConcurrentChange structure serialized (bytea). - * We can't store the tuple directly because tuplestore only supports - * minimum tuple and we may need to transfer OID system column from the - * output plugin. Also we need to transfer the change kind, so it's better - * to put everything in the structure than to use 2 tuplestores "in - * parallel". - */ - TupleDesc tupdesc_change; - - /* Tuple descriptor needed to update indexes. */ + /* Tuple descriptor of the relation being processed. */ TupleDesc tupdesc; - /* Slot to retrieve data from tstore. */ - TupleTableSlot *tsslot; - - ResourceOwner resowner; + /* The current output file. */ + BufFile *file; } RepackDecodingState; +extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; + +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); extern void cluster_rel(RepackCommand command, Relation OldHeap, Oid indexOid, @@ -136,6 +103,6 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, extern bool am_decoding_for_repack(void); extern bool change_useless_for_repack(XLogRecordBuffer *buf); -extern void repack_decode_concurrent_changes(LogicalDecodingContext *ctx, - XLogRecPtr end_of_wal); + +extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/storage/procsignal.h b/src/include/storage/procsignal.h index e52b8eb7697..3ef35ca6b80 100644 --- a/src/include/storage/procsignal.h +++ b/src/include/storage/procsignal.h @@ -36,6 +36,7 @@ typedef enum PROCSIG_BARRIER, /* global barrier interrupt */ PROCSIG_LOG_MEMORY_CONTEXT, /* ask backend to log the memory contexts */ PROCSIG_PARALLEL_APPLY_MESSAGE, /* Message from parallel apply workers */ + PROCSIG_REPACK_MESSAGE, /* Message from repack worker */ /* Recovery conflict reasons */ PROCSIG_RECOVERY_CONFLICT_FIRST, diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index a0b7b38a5e2..d1a694f9008 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -496,7 +496,6 @@ CompressFileHandle CompressionLocation CompressorState ComputeXidHorizonsResult -ConcurrentChange ConcurrentChangeKind ConditionVariable ConditionVariableMinimallyPadded @@ -636,6 +635,9 @@ DeclareCursorStmt DecodedBkpBlock DecodedXLogRecord DecodingOutputState +DecodingWorker +DecodingWorkerShared +DecodingWorkerState DefElem DefElemAction DefaultACLInfo -- 2.47.3 --=-=-= Content-Type: text/plain Content-Disposition: attachment; filename=v30-0006-Use-multiple-snapshots-to-copy-the-data.patch ^ permalink raw reply [nested|flat] 385+ messages in thread
end of thread, other threads:[~2026-01-12 16:30 UTC | newest] Thread overview: 385+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v32 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v34 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v28 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v29 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v33 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v30 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v31 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v29 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v37 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2023-05-31 11:58 [PATCH v38 09/11] Add support for min/max aggregates for IVM Yugo Nagata <[email protected]> 2026-01-12 16:30 [PATCH 5/6] Use background worker to do logical decoding. Antonin Houska <[email protected]>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox